agora inbox for pgsql-hackers@postgresql.orghelp / color / mirror / Atom feed
[PATCH v32 08/11] Add aggregates support in IVM 324+ messages / 2 participants [nested] [flat]
* [PATCH v32 08/11] Add aggregates support in IVM @ 2023-05-31 11:46 Yugo Nagata <nagata@sraoss.co.jp> 0 siblings, 0 replies; 324+ messages in thread From: Yugo Nagata @ 2023-05-31 11:46 UTC (permalink / raw) count, sum, adn avg are supported. As a restriction, expressions specified in GROUP BY must appear in the target list because tuples to be updated in IMMV are identified by using this group key. However, in the case of aggregates without GROUP BY, there is only one tuple in the view, so keys are not uses to identify tuples. When creating a IMMV, in addition to __ivm_count column, some hidden columns for each aggregate are added to the target list. For example, names of these hidden columns are ivm_count_avg and ivm_sum_avg for the average function, and so on. When a base table is modified, the aggregated values and related hidden columns are also updated as well as __ivm_count__. The way of update depends the kind of aggregate function.=E3=80=80Specifically, sum and count are updated by simply adding or subtracting delta value calculated from delta tables. avg is updated by using values of sum and count stored in views as hidden columns and deltas calculated from delta tables. About aggregate functions except "count()" (sum and avg), NULLs in input values are ignored, and the result of aggegate should be NULL when no rows are selected. To support this specification, the numbers of non-NULL input values are counted and stored in hidden columns. In the case of count(), count(x) returns zero when no rows are selected, but count(*) doesn't ignore NULL input. --- src/backend/commands/createas.c | 265 +++++++++++++++++-- src/backend/commands/matview.c | 433 ++++++++++++++++++++++++++++++-- src/include/commands/createas.h | 1 + 3 files changed, 662 insertions(+), 37 deletions(-) diff --git a/src/backend/commands/createas.c b/src/backend/commands/createa= s.c index 299c5a133c..ecec93ec1c 100644 --- a/src/backend/commands/createas.c +++ b/src/backend/commands/createas.c @@ -51,13 +51,19 @@ #include "parser/parsetree.h" #include "parser/parse_clause.h" #include "parser/parse_func.h" +#include "parser/parse_type.h" #include "rewrite/rewriteHandler.h" +#include "rewrite/rewriteManip.h" +#include "storage/smgr.h" #include "tcop/tcopprot.h" #include "utils/builtins.h" #include "utils/lsyscache.h" +#include "utils/regproc.h" +#include "utils/fmgroids.h" #include "utils/rel.h" #include "utils/rls.h" #include "utils/snapmgr.h" +#include "utils/syscache.h" =20 typedef struct { @@ -71,6 +77,11 @@ typedef struct BulkInsertState bistate; /* bulk insert state */ } DR_intorel; =20 +typedef struct +{ + bool has_agg; +} check_ivm_restriction_context; + /* utility functions for CTAS definition creation */ static ObjectAddress create_ctas_internal(List *attrList, IntoClause *into= ); static ObjectAddress create_ctas_nodata(List *tlist, IntoClause *into); @@ -85,8 +96,9 @@ static void CreateIvmTriggersOnBaseTablesRecurse(Query *q= ry, Node *node, Oid mat Relids *relids, bool ex_lock); static void CreateIvmTrigger(Oid relOid, Oid viewOid, int16 type, int16 ti= ming, bool ex_lock); static void check_ivm_restriction(Node *node); -static bool check_ivm_restriction_walker(Node *node, void *context); +static bool check_ivm_restriction_walker(Node *node, check_ivm_restriction= _context *context); static Bitmapset *get_primary_key_attnos_from_query(Query *query, List **c= onstraintList); +static bool check_aggregate_supports_ivm(Oid aggfnoid); =20 /* * create_ctas_internal @@ -417,6 +429,7 @@ ExecCreateTableAs(ParseState *pstate, CreateTableAsStmt= *stmt, * rewriteQueryForIMMV -- rewrite view definition query for IMMV * * count(*) is added for counting distinct tuples in views. + * Also, additional hidden columns are added for aggregate values. */ Query * rewriteQueryForIMMV(Query *query, List *colNames) @@ -430,16 +443,49 @@ rewriteQueryForIMMV(Query *query, List *colNames) rewritten =3D copyObject(query); pstate->p_expr_kind =3D EXPR_KIND_SELECT_TARGET; =20 - /* - * Convert DISTINCT to GROUP BY and add count(*) for counting distinct - * tuples in views. - */ - if (rewritten->distinctClause) + /* group keys must be in targetlist */ + if (rewritten->groupClause) { - TargetEntry *tle; + ListCell *lc; + foreach(lc, rewritten->groupClause) + { + SortGroupClause *scl =3D (SortGroupClause *) lfirst(lc); + TargetEntry *tle =3D get_sortgroupclause_tle(scl, rewritten->targetList= ); =20 + if (tle->resjunk) + ereport(ERROR, + (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), + errmsg("GROUP BY expression not appearing in select list is not sup= ported on incrementally maintainable materialized view"))); + } + } + /* Convert DISTINCT to GROUP BY. count(*) will be added afterward. */ + else if (!rewritten->hasAggs && rewritten->distinctClause) rewritten->groupClause =3D transformDistinctClause(NULL, &rewritten->tar= getList, rewritten->sortClause, false); =20 + /* Add additional columns for aggregate values */ + if (rewritten->hasAggs) + { + ListCell *lc; + List *aggs =3D NIL; + AttrNumber next_resno =3D list_length(rewritten->targetList) + 1; + + foreach(lc, rewritten->targetList) + { + TargetEntry *tle =3D (TargetEntry *) lfirst(lc); + char *resname =3D (colNames =3D=3D NIL || foreach_current_index(lc) >= =3D list_length(colNames) ? + tle->resname : strVal(list_nth(colNames, tle->resno - 1))); + + if (IsA(tle->expr, Aggref)) + makeIvmAggColumn(pstate, (Aggref *) tle->expr, resname, &next_resno, &= aggs); + } + rewritten->targetList =3D list_concat(rewritten->targetList, aggs); + } + + /* Add count(*) for counting distinct tuples in views */ + if (rewritten->distinctClause || rewritten->hasAggs) + { + TargetEntry *tle; + fn =3D makeFuncCall(SystemFuncName("count"), NIL, COERCE_EXPLICIT_CALL, = -1); fn->agg_star =3D true; =20 @@ -456,6 +502,91 @@ rewriteQueryForIMMV(Query *query, List *colNames) return rewritten; } =20 +/* + * makeIvmAggColumn -- make additional aggregate columns for IVM + * + * For an aggregate column specified by aggref, additional aggregate colum= ns + * are added, which are used to calculate the new aggregate value in IMMV. + * An additional aggregate columns has a name based on resname + * (ex. ivm_count_resname), and resno specified by next_resno. The created + * columns are returned to aggs, and the resno for the next column is also + * returned to next_resno. + * + * Currently, an additional count() is created for aggref other than count. + * In addition, sum() is created for avg aggregate column. + */ +void +makeIvmAggColumn(ParseState *pstate, Aggref *aggref, char *resname, AttrNu= mber *next_resno, List **aggs) +{ + TargetEntry *tle_count; + Node *node; + FuncCall *fn; + Const *dmy_arg =3D makeConst(INT4OID, + -1, + InvalidOid, + sizeof(int32), + Int32GetDatum(1), + false, + true); /* pass by value */ + const char *aggname =3D get_func_name(aggref->aggfnoid); + + /* + * For aggregate functions except count, add count() func with the same a= rg parameters. + * This count result is used for determining if the aggregate value shoul= d be NULL or not. + * Also, add sum() func for avg because we need to calculate an average v= alue as sum/count. + * + * XXX: If there are same expressions explicitly in the target list, we c= an use this instead + * of adding new duplicated one. + */ + if (strcmp(aggname, "count") !=3D 0) + { + fn =3D makeFuncCall(SystemFuncName("count"), NIL, COERCE_EXPLICIT_CALL, = -1); + + /* Make a Func with a dummy arg, and then override this by the original = agg's args. */ + node =3D ParseFuncOrColumn(pstate, fn->funcname, list_make1(dmy_arg), NU= LL, fn, false, -1); + ((Aggref *)node)->args =3D aggref->args; + + tle_count =3D makeTargetEntry((Expr *) node, + *next_resno, + pstrdup(makeObjectName("__ivm_count",resname, "_")), + false); + *aggs =3D lappend(*aggs, tle_count); + (*next_resno)++; + } + if (strcmp(aggname, "avg") =3D=3D 0) + { + List *dmy_args =3D NIL; + ListCell *lc; + foreach(lc, aggref->aggargtypes) + { + Oid typeid =3D lfirst_oid(lc); + Type type =3D typeidType(typeid); + + Const *con =3D makeConst(typeid, + -1, + typeTypeCollation(type), + typeLen(type), + (Datum) 0, + true, + typeByVal(type)); + dmy_args =3D lappend(dmy_args, con); + ReleaseSysCache(type); + } + fn =3D makeFuncCall(SystemFuncName("sum"), NIL, COERCE_EXPLICIT_CALL, -1= ); + + /* Make a Func with dummy args, and then override this by the original a= gg's args. */ + node =3D ParseFuncOrColumn(pstate, fn->funcname, dmy_args, NULL, fn, fal= se, -1); + ((Aggref *)node)->args =3D aggref->args; + + tle_count =3D makeTargetEntry((Expr *) node, + *next_resno, + pstrdup(makeObjectName("__ivm_sum",resname, "_")), + false); + *aggs =3D lappend(*aggs, tle_count); + (*next_resno)++; + } +} + /* * GetIntoRelEFlags --- compute executor flags needed for CREATE TABLE AS * @@ -941,11 +1072,13 @@ CreateIvmTrigger(Oid relOid, Oid viewOid, int16 type= , int16 timing, bool ex_lock static void check_ivm_restriction(Node *node) { - check_ivm_restriction_walker(node, NULL); + check_ivm_restriction_context context =3D {false}; + + check_ivm_restriction_walker(node, &context); } =20 static bool -check_ivm_restriction_walker(Node *node, void *context) +check_ivm_restriction_walker(Node *node, check_ivm_restriction_context *co= ntext) { if (node =3D=3D NULL) return false; @@ -974,6 +1107,10 @@ check_ivm_restriction_walker(Node *node, void *contex= t) ereport(ERROR, (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), errmsg("CTE is not supported on incrementally maintainable materia= lized view"))); + if (qry->groupClause !=3D NIL && !qry->hasAggs) + ereport(ERROR, + (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), + errmsg("GROUP BY clause without aggregate is not supported on incr= ementally maintainable materialized view"))); if (qry->havingQual !=3D NULL) ereport(ERROR, (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), @@ -1026,6 +1163,8 @@ check_ivm_restriction_walker(Node *node, void *contex= t) } } =20 + context->has_agg |=3D qry->hasAggs; + /* restrictions for rtable */ foreach(lc, qry->rtable) { @@ -1074,7 +1213,7 @@ check_ivm_restriction_walker(Node *node, void *contex= t) =20 } =20 - query_tree_walker(qry, check_ivm_restriction_walker, NULL, QTW_IGNORE_= RANGE_TABLE); + query_tree_walker(qry, check_ivm_restriction_walker, (void *) context,= QTW_IGNORE_RANGE_TABLE); =20 break; } @@ -1085,8 +1224,12 @@ check_ivm_restriction_walker(Node *node, void *conte= xt) ereport(ERROR, (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), errmsg("column name %s is not supported on incrementally maintain= able materialized view", tle->resname))); + if (context->has_agg && !IsA(tle->expr, Aggref) && contain_aggs_of_lev= el((Node *) tle->expr, 0)) + ereport(ERROR, + (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), + errmsg("expression containing an aggregate in it is not supported = on incrementally maintainable materialized view"))); =20 - expression_tree_walker(node, check_ivm_restriction_walker, NULL); + expression_tree_walker(node, check_ivm_restriction_walker, (void *) co= ntext); break; } case T_JoinExpr: @@ -1098,14 +1241,36 @@ check_ivm_restriction_walker(Node *node, void *cont= ext) (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), errmsg("OUTER JOIN is not supported on incrementally maintainable= materialized view"))); =20 - expression_tree_walker(node, check_ivm_restriction_walker, NULL); + expression_tree_walker(node, check_ivm_restriction_walker, (void *) co= ntext); + break; } - break; case T_Aggref: - ereport(ERROR, - (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), - errmsg("aggregate function is not supported on incrementally maintai= nable materialized view"))); - break; + { + /* Check if this supports IVM */ + Aggref *aggref =3D (Aggref *) node; + const char *aggname =3D format_procedure(aggref->aggfnoid); + + if (aggref->aggfilter !=3D NULL) + ereport(ERROR, + (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), + errmsg("aggregate function with FILTER clause is not supported on = incrementally maintainable materialized view"))); + + if (aggref->aggdistinct !=3D NULL) + ereport(ERROR, + (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), + errmsg("aggregate function with DISTINCT arguments is not supporte= d on incrementally maintainable materialized view"))); + + if (aggref->aggorder !=3D NULL) + ereport(ERROR, + (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), + errmsg("aggregate function with ORDER clause is not supported on i= ncrementally maintainable materialized view"))); + + if (!check_aggregate_supports_ivm(aggref->aggfnoid)) + ereport(ERROR, + (errcode(ERRCODE_FEATURE_NOT_SUPPORTED), + errmsg("aggregate function %s is not supported on incrementally ma= intainable materialized view", aggname))); + break; + } default: expression_tree_walker(node, check_ivm_restriction_walker, (void *) con= text); break; @@ -1113,6 +1278,46 @@ check_ivm_restriction_walker(Node *node, void *conte= xt) return false; } =20 +/* + * check_aggregate_supports_ivm + * + * Check if the given aggregate function is supporting IVM + */ +static bool +check_aggregate_supports_ivm(Oid aggfnoid) +{ + switch (aggfnoid) + { + /* count */ + case F_COUNT_ANY: + case F_COUNT_: + + /* sum */ + case F_SUM_INT8: + case F_SUM_INT4: + case F_SUM_INT2: + case F_SUM_FLOAT4: + case F_SUM_FLOAT8: + case F_SUM_MONEY: + case F_SUM_INTERVAL: + case F_SUM_NUMERIC: + + /* avg */ + case F_AVG_INT8: + case F_AVG_INT4: + case F_AVG_INT2: + case F_AVG_NUMERIC: + case F_AVG_FLOAT4: + case F_AVG_FLOAT8: + case F_AVG_INTERVAL: + + return true; + + default: + return false; + } +} + /* * CreateIndexOnIMMV * @@ -1170,7 +1375,29 @@ CreateIndexOnIMMV(Query *query, Relation matviewRel) index->concurrent =3D false; index->if_not_exists =3D false; =20 - if (query->distinctClause) + if (query->groupClause) + { + /* create unique constraint on GROUP BY expression columns */ + foreach(lc, query->groupClause) + { + SortGroupClause *scl =3D (SortGroupClause *) lfirst(lc); + TargetEntry *tle =3D get_sortgroupclause_tle(scl, query->targetList); + Form_pg_attribute attr =3D TupleDescAttr(matviewRel->rd_att, tle->resno= - 1); + IndexElem *iparam; + + iparam =3D makeNode(IndexElem); + iparam->name =3D pstrdup(NameStr(attr->attname)); + iparam->expr =3D NULL; + iparam->indexcolname =3D NULL; + iparam->collation =3D NIL; + iparam->opclass =3D NIL; + iparam->opclassopts =3D NIL; + iparam->ordering =3D SORTBY_DEFAULT; + iparam->nulls_ordering =3D SORTBY_NULLS_DEFAULT; + index->indexParams =3D lappend(index->indexParams, iparam); + } + } + else if (query->distinctClause) { /* create unique constraint on all columns */ foreach(lc, query->targetList) @@ -1228,7 +1455,7 @@ CreateIndexOnIMMV(Query *query, Relation matviewRel) (errmsg("could not create an index on materialized view \"%s\" automa= tically", RelationGetRelationName(matviewRel)), errdetail("This target list does not have all the primary key column= s, " - "or this view does not contain DISTINCT clause."), + "or this view does not contain GROUP BY or DISTINCT clause."), errhint("Create an index on the materialized view for efficient incr= emental maintenance."))); return; } diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c index 0064e10966..b7f6c3831b 100644 --- a/src/backend/commands/matview.c +++ b/src/backend/commands/matview.c @@ -27,6 +27,7 @@ #include "catalog/pg_trigger.h" #include "catalog/pg_opclass.h" #include "commands/cluster.h" +#include "commands/defrem.h" #include "commands/matview.h" #include "commands/tablecmds.h" #include "commands/tablespace.h" @@ -36,6 +37,7 @@ #include "executor/tstoreReceiver.h" #include "miscadmin.h" #include "nodes/makefuncs.h" +#include "optimizer/optimizer.h" #include "parser/analyze.h" #include "parser/parse_clause.h" #include "parser/parse_func.h" @@ -107,6 +109,13 @@ static HTAB *mv_trigger_info =3D NULL; =20 static bool in_delta_calculation =3D false; =20 +/* kind of IVM operation for the view */ +typedef enum +{ + IVM_ADD, + IVM_SUB +} IvmOp; + /* ENR name for materialized view delta */ #define NEW_DELTA_ENRNAME "new_delta" #define OLD_DELTA_ENRNAME "old_delta" @@ -138,7 +147,7 @@ static RangeTblEntry *get_prestate_rte(RangeTblEntry *r= te, MV_TriggerTable *tabl QueryEnvironment *queryEnv, Oid matviewid); static RangeTblEntry *replace_rte_with_delta(RangeTblEntry *rte, MV_Trigge= rTable *table, bool is_new, QueryEnvironment *queryEnv); -static Query *rewrite_query_for_counting(Query *query, ParseState *pstate); +static Query *rewrite_query_for_counting_and_aggregates(Query *query, Pars= eState *pstate); =20 static void calc_delta(MV_TriggerTable *table, int rte_index, Query *query, DestReceiver *dest_old, DestReceiver *dest_new, @@ -149,14 +158,27 @@ static Query *rewrite_query_for_postupdate_state(Quer= y *query, MV_TriggerTable * static void apply_delta(Oid matviewOid, Tuplestorestate *old_tuplestores, = Tuplestorestate *new_tuplestores, TupleDesc tupdesc_old, TupleDesc tupdesc_new, Query *query, bool use_count, char *count_colname); +static void append_set_clause_for_count(const char *resname, StringInfo bu= f_old, + StringInfo buf_new,StringInfo aggs_list); +static void append_set_clause_for_sum(const char *resname, StringInfo buf_= old, + StringInfo buf_new, StringInfo aggs_list); +static void append_set_clause_for_avg(const char *resname, StringInfo buf_= old, + StringInfo buf_new, StringInfo aggs_list, + const char *aggtype); +static char *get_operation_string(IvmOp op, const char *col, const char *a= rg1, const char *arg2, + const char* count_col, const char *castType); +static char *get_null_condition_string(IvmOp op, const char *arg1, const c= har *arg2, + const char* count_col); static void apply_old_delta(const char *matviewname, const char *deltaname= _old, List *keys); static void apply_old_delta_with_count(const char *matviewname, const char= *deltaname_old, - List *keys, const char *count_colname); + List *keys, StringInfo aggs_list, StringInfo aggs_set, + const char *count_colname); static void apply_new_delta(const char *matviewname, const char *deltaname= _new, StringInfo target_list); static void apply_new_delta_with_count(const char *matviewname, const char= * deltaname_new, - List *keys, StringInfo target_list, const char* count_colname); + List *keys, StringInfo target_list, StringInfo aggs_set, + const char* count_colname); static char *get_matching_condition_string(List *keys); static void generate_equal(StringInfo querybuf, Oid opttype, const char *leftop, const char *rightop); @@ -1454,11 +1476,44 @@ IVM_immediate_maintenance(PG_FUNCTION_ARGS) * When a base table is truncated, the view content will be empty if the * view definition query does not contain an aggregate without a GROUP cl= ause. * Therefore, such views can be truncated. + * + * Aggregate views without a GROUP clause always have one row. Therefore, + * if a base table is truncated, the view will not be empty and will cont= ain + * a row with NULL value (or 0 for count()). So, in this case, we refresh= the + * view instead of truncating it. */ if (TRIGGER_FIRED_BY_TRUNCATE(trigdata->tg_event)) { - ExecuteTruncateGuts(list_make1(matviewRel), list_make1_oid(matviewOid), - NIL, DROP_RESTRICT, false, false); + if (!(query->hasAggs && query->groupClause =3D=3D NIL)) + ExecuteTruncateGuts(list_make1(matviewRel), list_make1_oid(matviewOid), + NIL, DROP_RESTRICT, false, false); + else + { + Oid OIDNewHeap; + DestReceiver *dest; + uint64 processed =3D 0; + Query *dataQuery =3D rewriteQueryForIMMV(query, NIL); + char relpersistence =3D matviewRel->rd_rel->relpersistence; + + /* + * Create the transient table that will receive the regenerated data. L= ock + * it against access by any other process until commit (by which time it + * will be gone). + */ + OIDNewHeap =3D make_new_heap(matviewOid, matviewRel->rd_rel->reltablesp= ace, + matviewRel->rd_rel->relam, + relpersistence, ExclusiveLock); + LockRelationOid(OIDNewHeap, AccessExclusiveLock); + dest =3D CreateTransientRelDestReceiver(OIDNewHeap); + + /* Generate the data */ + processed =3D refresh_matview_datafill(dest, dataQuery, NULL, NULL, ""); + refresh_by_heap_swap(matviewOid, OIDNewHeap, relpersistence); + + /* Inform cumulative stats system about our activity */ + pgstat_count_truncate(matviewRel); + pgstat_count_heap_insert(matviewRel, processed); + } =20 /* Clean up hash entry and delete tuplestores */ clean_up_IVM_hash_entry(entry, false); @@ -1498,8 +1553,8 @@ IVM_immediate_maintenance(PG_FUNCTION_ARGS) /* Set all tables in the query to pre-update state */ rewritten =3D rewrite_query_for_preupdate_state(rewritten, entry->tables, pstate, matviewOid); - /* Rewrite for counting duplicated tuples */ - rewritten =3D rewrite_query_for_counting(rewritten, pstate); + /* Rewrite for counting duplicated tuples and aggregates functions*/ + rewritten =3D rewrite_query_for_counting_and_aggregates(rewritten, pstate= ); =20 /* Create tuplestores to store view deltas */ if (entry->has_old) @@ -1550,7 +1605,7 @@ IVM_immediate_maintenance(PG_FUNCTION_ARGS) =20 count_colname =3D pstrdup("__ivm_count__"); =20 - if (query->distinctClause) + if (query->hasAggs || query->distinctClause) use_count =3D true; =20 /* calculate delta tables */ @@ -1946,17 +2001,34 @@ replace_rte_with_delta(RangeTblEntry *rte, MV_Trigg= erTable *table, bool is_new, } =20 /* - * rewrite_query_for_counting + * rewrite_query_for_counting_and_aggregates * - * Rewrite query for counting duplicated tuples. + * Rewrite query for counting duplicated tuples and aggregate functions. */ static Query * -rewrite_query_for_counting(Query *query, ParseState *pstate) +rewrite_query_for_counting_and_aggregates(Query *query, ParseState *pstate) { TargetEntry *tle_count; FuncCall *fn; Node *node; =20 + /* For aggregate views */ + if (query->hasAggs) + { + ListCell *lc; + List *aggs =3D NIL; + AttrNumber next_resno =3D list_length(query->targetList) + 1; + + foreach(lc, query->targetList) + { + TargetEntry *tle =3D (TargetEntry *) lfirst(lc); + + if (IsA(tle->expr, Aggref)) + makeIvmAggColumn(pstate, (Aggref *)tle->expr, tle->resname, &next_resn= o, &aggs); + } + query->targetList =3D list_concat(query->targetList, aggs); + } + /* Add count(*) for counting distinct tuples in views */ fn =3D makeFuncCall(SystemFuncName("count"), NIL, COERCE_EXPLICIT_CALL, -= 1); fn->agg_star =3D true; @@ -2029,6 +2101,8 @@ rewrite_query_for_postupdate_state(Query *query, MV_T= riggerTable *table, int rte return query; } =20 +#define IVM_colname(type, col) makeObjectName("__ivm_" type, col, "_") + /* * apply_delta * @@ -2042,6 +2116,9 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tupl= estores, Tuplestorestate *n { StringInfoData querybuf; StringInfoData target_list_buf; + StringInfo aggs_list_buf =3D NULL; + StringInfo aggs_set_old =3D NULL; + StringInfo aggs_set_new =3D NULL; Relation matviewRel; char *matviewname; ListCell *lc; @@ -2064,6 +2141,15 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tup= lestores, Tuplestorestate *n initStringInfo(&querybuf); initStringInfo(&target_list_buf); =20 + if (query->hasAggs) + { + if (old_tuplestores && tuplestore_tuple_count(old_tuplestores) > 0) + aggs_set_old =3D makeStringInfo(); + if (new_tuplestores && tuplestore_tuple_count(new_tuplestores) > 0) + aggs_set_new =3D makeStringInfo(); + aggs_list_buf =3D makeStringInfo(); + } + /* build string of target list */ for (i =3D 0; i < matviewRel->rd_att->natts; i++) { @@ -2080,13 +2166,61 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tu= plestores, Tuplestorestate *n { TargetEntry *tle =3D (TargetEntry *) lfirst(lc); Form_pg_attribute attr =3D TupleDescAttr(matviewRel->rd_att, i); + char *resname =3D NameStr(attr->attname); =20 i++; =20 if (tle->resjunk) continue; =20 - keys =3D lappend(keys, attr); + /* + * For views without aggregates, all attributes are used as keys to iden= tify a + * tuple in a view. + */ + if (!query->hasAggs) + keys =3D lappend(keys, attr); + + /* For views with aggregates, we need to build SET clause for updating a= ggregate + * values. */ + if (query->hasAggs && IsA(tle->expr, Aggref)) + { + Aggref *aggref =3D (Aggref *) tle->expr; + const char *aggname =3D get_func_name(aggref->aggfnoid); + + /* + * We can use function names here because it is already checked if these + * can be used in IMMV by its OID at the definition time. + */ + + /* count */ + if (!strcmp(aggname, "count")) + append_set_clause_for_count(resname, aggs_set_old, aggs_set_new, aggs_= list_buf); + + /* sum */ + else if (!strcmp(aggname, "sum")) + append_set_clause_for_sum(resname, aggs_set_old, aggs_set_new, aggs_li= st_buf); + + /* avg */ + else if (!strcmp(aggname, "avg")) + append_set_clause_for_avg(resname, aggs_set_old, aggs_set_new, aggs_li= st_buf, + format_type_be(aggref->aggtype)); + + else + elog(ERROR, "unsupported aggregate function: %s", aggname); + } + } + + /* If we have GROUP BY clause, we use its entries as keys. */ + if (query->hasAggs && query->groupClause) + { + foreach (lc, query->groupClause) + { + SortGroupClause *sgcl =3D (SortGroupClause *) lfirst(lc); + TargetEntry *tle =3D get_sortgroupclause_tle(sgcl, query->targetList); + Form_pg_attribute attr =3D TupleDescAttr(matviewRel->rd_att, tle->resno= - 1); + + keys =3D lappend(keys, attr); + } } =20 /* Start maintaining the materialized view. */ @@ -2117,7 +2251,8 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tupl= estores, Tuplestorestate *n if (use_count) /* apply old delta and get rows to be recalculated */ apply_old_delta_with_count(matviewname, OLD_DELTA_ENRNAME, - keys, count_colname); + keys, aggs_list_buf, aggs_set_old, + count_colname); else apply_old_delta(matviewname, OLD_DELTA_ENRNAME, keys); =20 @@ -2143,7 +2278,7 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tupl= estores, Tuplestorestate *n /* apply new delta */ if (use_count) apply_new_delta_with_count(matviewname, NEW_DELTA_ENRNAME, - keys, &target_list_buf, count_colname); + keys, aggs_set_new, &target_list_buf, count_colname); else apply_new_delta(matviewname, NEW_DELTA_ENRNAME, &target_list_buf); } @@ -2158,6 +2293,250 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tu= plestores, Tuplestorestate *n elog(ERROR, "SPI_finish failed"); } =20 +/* + * append_set_clause_for_count + * + * Append SET clause string for count aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + */ +static void +append_set_clause_for_count(const char *resname, StringInfo buf_old, + StringInfo buf_new,StringInfo aggs_list) +{ + /* For tuple deletion */ + if (buf_old) + { + /* resname =3D mv.resname - t.resname */ + appendStringInfo(buf_old, + ", %s =3D %s", + quote_qualified_identifier(NULL, resname), + get_operation_string(IVM_SUB, resname, "mv", "t", NULL, NULL)); + } + /* For tuple insertion */ + if (buf_new) + { + /* resname =3D mv.resname + diff.resname */ + appendStringInfo(buf_new, + ", %s =3D %s", + quote_qualified_identifier(NULL, resname), + get_operation_string(IVM_ADD, resname, "mv", "diff", NULL, NULL)); + } + + appendStringInfo(aggs_list, ", %s", + quote_qualified_identifier("diff", resname) + ); +} + +/* + * append_set_clause_for_sum + * + * Append SET clause string for sum aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + */ +static void +append_set_clause_for_sum(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list) +{ + char *count_col =3D IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* sum =3D mv.sum - t.sum */ + appendStringInfo(buf_old, + ", %s =3D %s", + quote_qualified_identifier(NULL, resname), + get_operation_string(IVM_SUB, resname, "mv", "t", count_col, NULL) + ); + /* count =3D mv.count - t.count */ + appendStringInfo(buf_old, + ", %s =3D %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + } + /* For tuple insertion */ + if (buf_new) + { + /* sum =3D mv.sum + diff.sum */ + appendStringInfo(buf_new, + ", %s =3D %s", + quote_qualified_identifier(NULL, resname), + get_operation_string(IVM_ADD, resname, "mv", "diff", count_col, NULL) + ); + /* count =3D mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s =3D %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + +/* + * append_set_clause_for_avg + * + * Append SET clause string for avg aggregation to given buffers. + * Also, append resnames required for calculating the aggregate value. + */ +static void +append_set_clause_for_avg(const char *resname, StringInfo buf_old, + StringInfo buf_new, StringInfo aggs_list, + const char *aggtype) +{ + char *sum_col =3D IVM_colname("sum", resname); + char *count_col =3D IVM_colname("count", resname); + + /* For tuple deletion */ + if (buf_old) + { + /* avg =3D (mv.sum - t.sum)::aggtype / (mv.count - t.count) */ + appendStringInfo(buf_old, + ", %s =3D %s OPERATOR(pg_catalog./) %s", + quote_qualified_identifier(NULL, resname), + get_operation_string(IVM_SUB, sum_col, "mv", "t", count_col, aggtype), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + /* sum =3D mv.sum - t.sum */ + appendStringInfo(buf_old, + ", %s =3D %s", + quote_qualified_identifier(NULL, sum_col), + get_operation_string(IVM_SUB, sum_col, "mv", "t", count_col, NULL) + ); + /* count =3D mv.count - t.count */ + appendStringInfo(buf_old, + ", %s =3D %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_SUB, count_col, "mv", "t", NULL, NULL) + ); + + } + /* For tuple insertion */ + if (buf_new) + { + /* avg =3D (mv.sum + diff.sum)::aggtype / (mv.count + diff.count) */ + appendStringInfo(buf_new, + ", %s =3D %s OPERATOR(pg_catalog./) %s", + quote_qualified_identifier(NULL, resname), + get_operation_string(IVM_ADD, sum_col, "mv", "diff", count_col, aggtype= ), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + /* sum =3D mv.sum + diff.sum */ + appendStringInfo(buf_new, + ", %s =3D %s", + quote_qualified_identifier(NULL, sum_col), + get_operation_string(IVM_ADD, sum_col, "mv", "diff", count_col, NULL) + ); + /* count =3D mv.count + diff.count */ + appendStringInfo(buf_new, + ", %s =3D %s", + quote_qualified_identifier(NULL, count_col), + get_operation_string(IVM_ADD, count_col, "mv", "diff", NULL, NULL) + ); + } + + appendStringInfo(aggs_list, ", %s, %s, %s", + quote_qualified_identifier("diff", resname), + quote_qualified_identifier("diff", IVM_colname("sum", resname)), + quote_qualified_identifier("diff", IVM_colname("count", resname)) + ); +} + +/* + * get_operation_string + * + * Build a string to calculate the new aggregate values. + */ +static char * +get_operation_string(IvmOp op, const char *col, const char *arg1, const ch= ar *arg2, + const char* count_col, const char *castType) +{ + StringInfoData buf; + StringInfoData castString; + char *col1 =3D quote_qualified_identifier(arg1, col); + char *col2 =3D quote_qualified_identifier(arg2, col); + char op_char =3D (op =3D=3D IVM_SUB ? '-' : '+'); + + initStringInfo(&buf); + initStringInfo(&castString); + + if (castType) + appendStringInfo(&castString, "::%s", castType); + + if (!count_col) + { + /* + * If the attributes don't have count columns then calc the result + * by using the operator simply. + */ + appendStringInfo(&buf, "(%s OPERATOR(pg_catalog.%c) %s)%s", + col1, op_char, col2, castString.data); + } + else + { + /* + * If the attributes have count columns then consider the condition + * where the result becomes NULL. + */ + char *null_cond =3D get_null_condition_string(op, arg1, arg2, count_col); + + appendStringInfo(&buf, + "(CASE WHEN %s THEN NULL " + "WHEN %s IS NULL THEN %s " + "WHEN %s IS NULL THEN %s " + "ELSE (%s OPERATOR(pg_catalog.%c) %s)%s END)", + null_cond, + col1, col2, + col2, col1, + col1, op_char, col2, castString.data + ); + } + + return buf.data; +} + +/* + * get_null_condition_string + * + * Build a predicate string for CASE clause to check if an aggregate value + * will became NULL after the given operation is applied. + */ +static char * +get_null_condition_string(IvmOp op, const char *arg1, const char *arg2, + const char* count_col) +{ + StringInfoData null_cond; + initStringInfo(&null_cond); + + switch (op) + { + case IVM_ADD: + appendStringInfo(&null_cond, + "%s OPERATOR(pg_catalog.=3D) 0 AND %s OPERATOR(pg_catalog.=3D) 0", + quote_qualified_identifier(arg1, count_col), + quote_qualified_identifier(arg2, count_col) + ); + break; + case IVM_SUB: + appendStringInfo(&null_cond, + "%s OPERATOR(pg_catalog.=3D) %s", + quote_qualified_identifier(arg1, count_col), + quote_qualified_identifier(arg2, count_col) + ); + break; + default: + elog(ERROR,"unknown operation"); + } + + return null_cond.data; +} + + /* * apply_old_delta_with_count * @@ -2165,13 +2544,20 @@ apply_delta(Oid matviewOid, Tuplestorestate *old_tu= plestores, Tuplestorestate *n * which contains tuples to be deleted from to a materialized view given by * matviewname. This is used when counting is required, that is, the view * has aggregate or distinct. + * + * If the view desn't have aggregates or has GROUP BY, this requires a keys + * list to identify a tuple in the view. If the view has aggregates, this + * requires strings representing resnames of aggregates and SET clause for + * updating aggregate values. */ static void apply_old_delta_with_count(const char *matviewname, const char *deltaname_= old, - List *keys, const char *count_colname) + List *keys, StringInfo aggs_list, StringInfo aggs_set, + const char *count_colname) { StringInfoData querybuf; char *match_cond; + bool agg_without_groupby =3D (list_length(keys) =3D=3D 0); =20 /* build WHERE condition for searching tuples to be deleted */ match_cond =3D get_matching_condition_string(keys); @@ -2181,22 +2567,26 @@ apply_old_delta_with_count(const char *matviewname,= const char *deltaname_old, appendStringInfo(&querybuf, "WITH t AS (" /* collecting tid of target tuples in the view */ "SELECT diff.%s, " /* count column */ - "(diff.%s OPERATOR(pg_catalog.=3D) mv.%s) AS for_dlt, " + "(diff.%s OPERATOR(pg_catalog.=3D) mv.%s AND %s) AS for_dlt, " "mv.ctid " + "%s " /* aggregate columns */ "FROM %s AS mv, %s AS diff " "WHERE %s" /* tuple matching condition */ "), updt AS (" /* update a tuple if this is not to be deleted */ "UPDATE %s AS mv SET %s =3D mv.%s OPERATOR(pg_catalog.-) t.%s " + "%s" /* SET clauses for aggregates */ "FROM t WHERE mv.ctid OPERATOR(pg_catalog.=3D) t.ctid AND NOT for_dl= t " ")" /* delete a tuple if this is to be deleted */ "DELETE FROM %s AS mv USING t " "WHERE mv.ctid OPERATOR(pg_catalog.=3D) t.ctid AND for_dlt", count_colname, - count_colname, count_colname, + count_colname, count_colname, (agg_without_groupby ? "false" : "true"= ), + (aggs_list !=3D NULL ? aggs_list->data : ""), matviewname, deltaname_old, match_cond, matviewname, count_colname, count_colname, count_colname, + (aggs_set !=3D NULL ? aggs_set->data : ""), matviewname); =20 if (SPI_exec(querybuf.data, 0) !=3D SPI_OK_DELETE) @@ -2260,10 +2650,15 @@ apply_old_delta(const char *matviewname, const char= *deltaname_old, * matviewname. This is used when counting is required, that is, the view * has aggregate or distinct. Also, when a table in EXISTS sub queries * is modified. + * + * If the view desn't have aggregates or has GROUP BY, this requires a keys + * list to identify a tuple in the view. If the view has aggregates, this + * requires strings representing SET clause for updating aggregate values. */ static void apply_new_delta_with_count(const char *matviewname, const char* deltaname_= new, - List *keys, StringInfo target_list, const char* count_colname) + List *keys, StringInfo aggs_set, StringInfo target_list, + const char* count_colname) { StringInfoData querybuf; StringInfoData returning_keys; @@ -2294,6 +2689,7 @@ apply_new_delta_with_count(const char *matviewname, c= onst char* deltaname_new, appendStringInfo(&querybuf, "WITH updt AS (" /* update a tuple if this exists in the view */ "UPDATE %s AS mv SET %s =3D mv.%s OPERATOR(pg_catalog.+) diff.%s " + "%s " /* SET clauses for aggregates */ "FROM %s AS diff " "WHERE %s " /* tuple matching condition */ "RETURNING %s" /* returning keys of updated tuples */ @@ -2301,6 +2697,7 @@ apply_new_delta_with_count(const char *matviewname, c= onst char* deltaname_new, "SELECT %s FROM %s AS diff " "WHERE NOT EXISTS (SELECT 1 FROM updt AS mv WHERE %s);", matviewname, count_colname, count_colname, count_colname, + (aggs_set !=3D NULL ? aggs_set->data : ""), deltaname_new, match_cond, returning_keys.data, diff --git a/src/include/commands/createas.h b/src/include/commands/createa= s.h index 6b47e66bfd..af3a5b4b27 100644 --- a/src/include/commands/createas.h +++ b/src/include/commands/createas.h @@ -30,6 +30,7 @@ extern void CreateIvmTriggersOnBaseTables(Query *qry, Oid= matviewOid); extern void CreateIndexOnIMMV(Query *query, Relation matviewRel); =20 extern Query *rewriteQueryForIMMV(Query *query, List *colNames); +extern void makeIvmAggColumn(ParseState *pstate, Aggref *aggref, char *res= name, AttrNumber *next_resno, List **aggs); =20 extern int GetIntoRelEFlags(IntoClause *intoClause); =20 --=20 2.25.1 --Multipart=_Sun__31_Mar_2024_22_59_31_+0900_msknEviJj08_wgqO Content-Type: text/x-diff; name="v32-0009-Add-support-for-min-max-aggregates-for-IVM.patch" Content-Disposition: attachment; filename="v32-0009-Add-support-for-min-max-aggregates-for-IVM.patch" Content-Transfer-Encoding: 7bit ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
* [PATCH v44 07/10] Split cluster.h to create repack_internal.h @ 2026-03-24 14:09 Álvaro Herrera <alvherre@kurilemu.de> 0 siblings, 0 replies; 324+ messages in thread From: Álvaro Herrera @ 2026-03-24 14:09 UTC (permalink / raw) Most of the details of concurrent repack are only needed by pgoutput_repack; and they have a nasty effect on headers included by cluster.h (used by several high-profile places), so isolate that for cleanliness. Also, change_useless_for_repack() is better declared in decode.h. --- src/backend/commands/cluster.c | 128 ++++++++++-------- src/backend/postmaster/bgworker.c | 2 +- .../pgoutput_repack/pgoutput_repack.c | 13 +- src/include/commands/cluster.h | 57 +------- src/include/commands/repack_internal.h | 59 ++++++++ src/include/replication/decode.h | 4 + 6 files changed, 147 insertions(+), 116 deletions(-) create mode 100644 src/include/commands/repack_internal.h diff --git a/src/backend/commands/cluster.c b/src/backend/commands/cluster.c index 75556cbdafb..2c3058ba10d 100644 --- a/src/backend/commands/cluster.c +++ b/src/backend/commands/cluster.c @@ -59,6 +59,7 @@ #include "commands/cluster.h" #include "commands/defrem.h" #include "commands/progress.h" +#include "commands/repack_internal.h" #include "commands/tablecmds.h" #include "commands/vacuum.h" #include "executor/executor.h" @@ -205,17 +206,11 @@ typedef struct DecodingWorkerShared char error_queue[FLEXIBLE_ARRAY_MEMBER]; } DecodingWorkerShared; -/* - * Generate worker's output file name. If relations of the same 'relid' happen - * to be processed at the same time, they must be from different databases and - * therefore different backends must be involved. (PID is already present in - * the fileset name.) - */ -static inline void -DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) -{ - snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); -} +/* Is this process a REPACK worker? */ +static bool is_repack_worker = false; + +static pid_t backend_pid; +static ProcNumber backend_proc_number; /* * Backend-local information to control the decoding worker. @@ -241,6 +236,7 @@ static DecodingWorker *decoding_worker = NULL; */ volatile sig_atomic_t RepackMessagePending = false; +static LOCKMODE RepackLockLevel(bool concurrent); static bool cluster_rel_recheck(RepackCommand cmd, Relation OldHeap, Oid indexOid, Oid userid, LOCKMODE lmode, int options); @@ -298,12 +294,16 @@ static Relation process_single_relation(RepackStmt *stmt, ClusterParams *params); static Oid determine_clustered_index(Relation rel, bool usingindex, const char *indexname); -static void start_decoding_worker(Oid relid); -static void stop_decoding_worker(void); + +static void start_repack_decoding_worker(Oid relid); +static void stop_repack_decoding_worker(void); static void repack_worker_internal(dsm_segment *seg); static void export_initial_snapshot(Snapshot snapshot, DecodingWorkerShared *shared); static Snapshot get_initial_snapshot(DecodingWorker *worker); +extern bool am_decoding_for_repack(void); +static void DecodingWorkerFileName(char *fname, Oid relid, uint32 seq); + static void ProcessRepackMessage(StringInfo msg); static const char *RepackCommandAsString(RepackCommand cmd); @@ -369,17 +369,8 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) parser_errposition(pstate, opt->location)); } - /* - * Determine the lock mode expected by cluster_rel(). - * - * In the exclusive case, we obtain AccessExclusiveLock right away to - * avoid lock-upgrade hazard in the single-transaction case. In the - * CONCURRENTLY case, the AccessExclusiveLock will only be used at the end - * of processing, supposedly for very short time. Until then, we'll have - * to unlock the relation temporarily, so there's no lock-upgrade hazard. - */ - lockmode = (params.options & CLUOPT_CONCURRENT) == 0 ? - AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lockmode = RepackLockLevel((params.options & CLUOPT_CONCURRENT) != 0); /* * If a single relation is specified, process it and we're done ... unless @@ -434,6 +425,12 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) "Repack", ALLOCSET_DEFAULT_SIZES); + /* + * Since we open a new transaction for each relation, we have to check + * that the relation still is what we think it is. + * + * In single-transaction CLUSTER, we don't need the overhead. + */ params.options |= CLUOPT_RECHECK; /* @@ -544,6 +541,22 @@ ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel) MemoryContextDelete(repack_context); } +/* + * In the non-concurrent case, we obtain AccessExclusiveLock throughout the + * operation to avoid any lock-upgrade hazards. In the concurrent case, we + * grab ShareUpdateExclusiveLock (jsut like VACUUM) for most of the + * processing and only acquire AccessExclusiveLock at the end, to swap the + * relation -- supposedly for a short time. + */ +static LOCKMODE +RepackLockLevel(bool concurrent) +{ + if (concurrent) + return ShareUpdateExclusiveLock; + else + return AccessExclusiveLock; +} + /* * cluster_rel * @@ -583,25 +596,22 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, bool concurrent = ((params->options & CLUOPT_CONCURRENT) != 0); Oid ident_idx = InvalidOid; - /* - * The lock mode is AccessExclusiveLock for normal processing and - * ShareUpdateExclusiveLock for concurrent processing (so that SELECT, - * INSERT, UPDATE and DELETE commands work, but cluster_rel() cannot be - * called concurrently for the same relation). - */ - lmode = !concurrent ? AccessExclusiveLock : ShareUpdateExclusiveLock; + /* Determine the lock mode to use. */ + lmode = RepackLockLevel(concurrent); - /* There are specific requirements on concurrent processing. */ + /* + * Check some preconditions in the concurrent case. This also obtains the + * replica index OID. + */ if (concurrent) { /* - * Make sure we have no XID assigned, otherwise call of - * repack_setup_logical_decoding() can cause a deadlock. + * Make sure we're not in a transaction block. * - * The existence of transaction block actually does not imply that XID - * was already assigned, but it very likely is. We might want to check - * the result of GetCurrentTransactionIdIfAny() instead, but that - * would be less clear from user's perspective. + * The reason is that repack_setup_logical_decoding() could deadlock + * if there's an XID already assigned. It would be possible to run in + * a transaction block if we had no XID, but this restriction is + * simpler for users to understand and we don't lose anything. */ PreventInTransactionBlock(isTopLevel, "REPACK (CONCURRENTLY)"); @@ -626,15 +636,11 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, RestrictSearchPath(); /* - * Since we may open a new transaction for each relation, we have to check - * that the relation still is what we think it is. + * Recheck that the relation is still what it was when we started. * - * If this is a single-transaction CLUSTER, we can skip these tests. We - * *must* skip the one on indisclustered since it would reject an attempt - * to cluster a not-previously-clustered index. - * - * XXX move [some of] these comments to where the RECHECK flag is - * determined? + * Note that it's critical to skip this in single-relation CLUSTER; + * otherwise, we would reject an attempt to cluster using a + * not-previously-clustered index. */ if (recheck && !cluster_rel_recheck(cmd, OldHeap, indexOid, save_userid, @@ -754,7 +760,7 @@ cluster_rel(RepackCommand cmd, Relation OldHeap, Oid indexOid, * However it still seems a good practice to make sure that the * worker never survives the REPACK command. */ - stop_decoding_worker(); + stop_repack_decoding_worker(); } } PG_END_TRY(); @@ -1093,7 +1099,7 @@ rebuild_relation(Relation OldHeap, Relation index, bool verbose, * clustering index) and checking again if it's still eligible for * REPACK CONCURRENTLY. */ - start_decoding_worker(tableOid); + start_repack_decoding_worker(tableOid); /* * Wait until the worker has the initial snapshot and retrieve it. @@ -2538,7 +2544,6 @@ RepackCommandAsString(RepackCommand cmd) return "???"; /* keep compiler quiet */ } - /* * Is this backend performing logical decoding on behalf of REPACK * (CONCURRENTLY) ? @@ -3688,7 +3693,7 @@ build_new_indexes(Relation NewHeap, Relation OldHeap, List *OldIndexes) * contents to a new table. */ static void -start_decoding_worker(Oid relid) +start_repack_decoding_worker(Oid relid) { Size size; dsm_segment *seg; @@ -3779,7 +3784,7 @@ start_decoding_worker(Oid relid) * we need to stop it explicitly at least on ERROR in the launching backend. */ static void -stop_decoding_worker(void) +stop_repack_decoding_worker(void) { BgwHandleStatus status; @@ -3817,12 +3822,6 @@ stop_decoding_worker(void) decoding_worker = NULL; } -/* Is this process a REPACK worker? */ -static bool is_repack_worker = false; - -static pid_t backend_pid; -static ProcNumber backend_proc_number; - /* * See ParallelWorkerShutdown for details. */ @@ -3869,7 +3868,7 @@ RepackWorkerMain(Datum main_arg) /* * Join locking group - see the comments around the call of - * start_decoding_worker(). + * start_repack_decoding_worker(). */ if (!BecomeLockGroupMember(shared->backend_proc, backend_pid)) /* The leader is not running anymore. */ @@ -4039,6 +4038,18 @@ get_initial_snapshot(DecodingWorker *worker) return snapshot; } +/* + * Generate worker's output file name. If relations of the same 'relid' happen + * to be processed at the same time, they must be from different databases and + * therefore different backends must be involved. (PID is already present in + * the fileset name.) + */ +static void +DecodingWorkerFileName(char *fname, Oid relid, uint32 seq) +{ + snprintf(fname, MAXPGPATH, "%u-%u", relid, seq); +} + bool IsRepackWorker(void) { @@ -4067,7 +4078,6 @@ void ProcessRepackMessages(void) { MemoryContext oldcontext; - static MemoryContext hpm_context = NULL; /* diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c index f8a8d1681e9..9e876d55e27 100644 --- a/src/backend/postmaster/bgworker.c +++ b/src/backend/postmaster/bgworker.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/parallel.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "libpq/pqsignal.h" #include "miscadmin.h" #include "pgstat.h" diff --git a/src/backend/replication/pgoutput_repack/pgoutput_repack.c b/src/backend/replication/pgoutput_repack/pgoutput_repack.c index 032fbd0e5b0..cc9ce615b18 100644 --- a/src/backend/replication/pgoutput_repack/pgoutput_repack.c +++ b/src/backend/replication/pgoutput_repack/pgoutput_repack.c @@ -13,7 +13,7 @@ #include "postgres.h" #include "access/detoast.h" -#include "commands/cluster.h" +#include "commands/repack_internal.h" #include "replication/snapbuild.h" #include "utils/memutils.h" @@ -176,7 +176,7 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, ConcurrentChangeKind kind, HeapTuple tuple) { RepackDecodingState *dstate; - MemoryContext oldcxt; + MemoryContext oldcxt; BufFile *file; List *attrs_ext = NIL; int natt_ext; @@ -226,7 +226,10 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, slot_getsomeattrs(slot, i + 1); - /* This is a non-null varlena datum, but we only care if it's out-of-line */ + /* + * This is a non-null varlena datum, but we only care if it's + * out-of-line + */ varlen = (varlena *) DatumGetPointer(slot->tts_values[i]); if (!VARATT_IS_EXTERNAL(varlen)) continue; @@ -244,8 +247,8 @@ repack_store_change(LogicalDecodingContext *ctx, Relation relation, * attributes (those actually should never appear on disk), so * only TOASTed attribute can be seen here. * - * FIXME in what circumstances can an ONDISK attr appear? - * Why aren't these written separately? + * FIXME in what circumstances can an ONDISK attr appear? Why + * aren't these written separately? */ Assert(VARATT_IS_EXTERNAL_ONDISK(varlen)); } diff --git a/src/include/commands/cluster.h b/src/include/commands/cluster.h index 1c0ac3ab4f5..1528d34fa42 100644 --- a/src/include/commands/cluster.h +++ b/src/include/commands/cluster.h @@ -13,17 +13,12 @@ #ifndef CLUSTER_H #define CLUSTER_H -#include "nodes/execnodes.h" +#include <signal.h> + #include "nodes/parsenodes.h" #include "parser/parse_node.h" -#include "replication/decode.h" -#include "postmaster/bgworker.h" -#include "replication/logical.h" -#include "storage/buffile.h" #include "storage/lockdefs.h" -#include "storage/shm_mq.h" #include "utils/relcache.h" -#include "utils/resowner.h" /* flag bits for ClusterParams->options */ @@ -34,55 +29,14 @@ #define CLUOPT_ANALYZE 0x08 /* do an ANALYZE */ #define CLUOPT_CONCURRENT 0x10 /* allow concurrent data changes */ - /* options for CLUSTER */ typedef struct ClusterParams { bits32 options; /* bitmask of CLUOPT_* */ } ClusterParams; - -/* - * The following definitions are used by REPACK CONCURRENTLY. - */ - -/* - * Stored as a single byte in the output file. - */ -#define CHANGE_INSERT 'i' -#define CHANGE_UPDATE_OLD 'u' -#define CHANGE_UPDATE_NEW 'U' -#define CHANGE_DELETE 'd' -typedef char ConcurrentChangeKind; - -/* - * Logical decoding state. - * - * The output plugin uses it to store the data changes that it decodes from - * WAL while the table contents is being copied to a new storage. - */ -typedef struct RepackDecodingState -{ -#ifdef USE_ASSERT_CHECKING - /* The relation whose changes we're decoding. */ - Oid relid; -#endif - - /* Per-change memory context. */ - MemoryContext change_cxt; - - /* A tuple slot used to pass tuples back and forth */ - TupleTableSlot *slot; - - /* The current output file. */ - BufFile *file; -} RepackDecodingState; - extern PGDLLIMPORT volatile sig_atomic_t RepackMessagePending; -extern bool IsRepackWorker(void); -extern void HandleRepackMessageInterrupt(void); -extern void ProcessRepackMessages(void); extern void ExecRepack(ParseState *pstate, RepackStmt *stmt, bool isTopLevel); @@ -104,8 +58,9 @@ extern void finish_heap_swap(Oid OIDOldHeap, Oid OIDNewHeap, MultiXactId cutoffMulti, char newrelpersistence); -extern bool am_decoding_for_repack(void); -extern bool change_useless_for_repack(XLogRecordBuffer *buf); +extern bool IsRepackWorker(void); +extern void HandleRepackMessageInterrupt(void); +extern void ProcessRepackMessages(void); + -extern void RepackWorkerMain(Datum main_arg); #endif /* CLUSTER_H */ diff --git a/src/include/commands/repack_internal.h b/src/include/commands/repack_internal.h new file mode 100644 index 00000000000..f90c973f5a0 --- /dev/null +++ b/src/include/commands/repack_internal.h @@ -0,0 +1,59 @@ +/*------------------------------------------------------------------------- + * + * repack_internal.h + * header for REPACK internals + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * Portions Copyright (c) 1994-5, Regents of the University of California + * + * src/include/commands/repack_internal.h + * + *------------------------------------------------------------------------- + */ +#ifndef REPACK_INTERNAL_H +#define REPACK_INTERNAL_H + +#include "nodes/execnodes.h" +#include "replication/decode.h" +#include "postmaster/bgworker.h" +#include "replication/logical.h" +#include "storage/buffile.h" +#include "storage/shm_mq.h" +#include "utils/resowner.h" + +/* + * Stored as a single byte in the output file. + */ +#define CHANGE_INSERT 'i' +#define CHANGE_UPDATE_OLD 'u' +#define CHANGE_UPDATE_NEW 'U' +#define CHANGE_DELETE 'd' +typedef char ConcurrentChangeKind; + +/* + * Logical decoding state. + * + * The output plugin uses it to store the data changes that it decodes from + * WAL while the table contents is being copied to a new storage. + */ +typedef struct RepackDecodingState +{ +#ifdef USE_ASSERT_CHECKING + /* The relation whose changes we're decoding. */ + Oid relid; +#endif + + /* Per-change memory context. */ + MemoryContext change_cxt; + + /* A tuple slot used to pass tuples back and forth */ + TupleTableSlot *slot; + + /* The current output file. */ + BufFile *file; +} RepackDecodingState; + + +extern void RepackWorkerMain(Datum main_arg); + +#endif /* REPACK_INTERNAL_H */ diff --git a/src/include/replication/decode.h b/src/include/replication/decode.h index 49f00fc48b8..02b5393474c 100644 --- a/src/include/replication/decode.h +++ b/src/include/replication/decode.h @@ -31,4 +31,8 @@ extern void logicalmsg_decode(LogicalDecodingContext *ctx, XLogRecordBuffer *buf extern void LogicalDecodingProcessRecord(LogicalDecodingContext *ctx, XLogReaderState *record); +/* in commands/cluster.c */ +extern bool change_useless_for_repack(XLogRecordBuffer *buf); + + #endif -- 2.47.3 --gwom7bl7ogtszo4k Content-Type: text/x-diff; charset=utf-8 Content-Disposition: attachment; filename="v44-0008-Use-BulkInsertState-when-copying-data-to-the-new.patch" ^ permalink raw reply [nested|flat] 324+ messages in thread
end of thread, other threads:[~2026-03-24 14:09 UTC | newest] Thread overview: 324+ messages (download: mbox mbox.gz follow: Atom feed) -- links below jump to the message on this page -- 2023-05-31 11:46 [PATCH v32 08/11] Add aggregates support in IVM Yugo Nagata <nagata@sraoss.co.jp> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de> 2026-03-24 14:09 [PATCH v44 07/10] Split cluster.h to create repack_internal.h Álvaro Herrera <alvherre@kurilemu.de>
This inbox is served by agora; see mirroring instructions for how to clone and mirror all data and code used for this inbox